agora inbox for pgsql-hackers@postgresql.orghelp / color / mirror / Atom feed
[PATCH 4/6] Fix obvious problems around the line datatype 940+ messages / 2 participants [nested] [flat]
* [PATCH 4/6] Fix obvious problems around the line datatype @ 2018-07-26 13:43 Tomas Vondra <tomas@2ndquadrant.com> 0 siblings, 0 replies; 940+ messages in thread From: Tomas Vondra @ 2018-07-26 13:43 UTC (permalink / raw) I have noticed some line operators retuning wrong results, and Tom Lane spotted similar problems on more places. Source history reveals that during 1990s, the internal format of the line datatype is changed, but most functions haven't got the hint. The fixes include: * Reject invalid specification A=B=0 on receive * Reject same points on line_construct_pp() * Fix perpendicular operator when negative values are involved * Avoid division by zero on perpendicular operator * Fix intersection and distance operators when neither A nor B are 1 * Return NULL for closest point when objects are parallel * Check whether closest point of line segments is the intersection point * Fix closest point of line segments being on the wrong segment The changes are also aiming make line operators more symmetric and less sensitive to floating point precision loss. The EPSILON interferes with every minor change in different ways. It is hard to guess which behaviour is more expected, but we can assume threating both sides of the operators more equally is more expected. Discussion: https://www.postgresql.org/message-id/CAE2gYzxF7-5djV6-cEvqQu-fNsnt%3DEqbOURx7ZDg%2BVv6ZMTWbg%40mail... Parallel discussions: * https://www.postgresql.org/message-id/CAE2gYzw_-z%3DV2kh8QqFjenu%3D8MJXzOP44wRW%3DAzzeamrmTT1%3DQ%40... * https://www.postgresql.org/message-id/20180201.205138.34583581.horiguchi.kyotaro@lab.ntt.co.jp --- src/backend/utils/adt/geo_ops.c | 153 +++++++++++++++++++++++++++------------- 1 file changed, 103 insertions(+), 50 deletions(-) diff --git a/src/backend/utils/adt/geo_ops.c b/src/backend/utils/adt/geo_ops.c index 314e442f2e..a72a9a28c9 100644 --- a/src/backend/utils/adt/geo_ops.c +++ b/src/backend/utils/adt/geo_ops.c @@ -48,6 +48,7 @@ static int point_inside(Point *p, int npts, Point *plist); /* Routines for lines */ static inline void line_construct(LINE *result, Point *pt, float8 m); +static inline float8 line_sl(LINE *line); static inline float8 line_invsl(LINE *line); static bool line_interpt_line(Point *result, LINE *l1, LINE *l2); static bool line_contain_point(LINE *line, Point *point); @@ -980,6 +981,11 @@ line_recv(PG_FUNCTION_ARGS) line->B = pq_getmsgfloat8(buf); line->C = pq_getmsgfloat8(buf); + if (FPzero(line->A) && FPzero(line->B)) + ereport(ERROR, + (errcode(ERRCODE_INVALID_BINARY_REPRESENTATION), + errmsg("invalid line specification: A and B cannot both be zero"))); + PG_RETURN_LINE_P(line); } @@ -1037,6 +1043,11 @@ line_construct_pp(PG_FUNCTION_ARGS) Point *pt2 = PG_GETARG_POINT_P(1); LINE *result = (LINE *) palloc(sizeof(LINE)); + if (point_eq_point(pt1, pt2)) + ereport(ERROR, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("invalid line specification: must be two distinct points"))); + line_construct(result, pt1, point_sl(pt1, pt2)); PG_RETURN_LINE_P(result); @@ -1073,11 +1084,15 @@ line_perp(PG_FUNCTION_ARGS) if (FPzero(l1->A)) PG_RETURN_BOOL(FPzero(l2->B)); - else if (FPzero(l1->B)) + if (FPzero(l2->A)) + PG_RETURN_BOOL(FPzero(l1->B)); + if (FPzero(l1->B)) PG_RETURN_BOOL(FPzero(l2->A)); + if (FPzero(l2->B)) + PG_RETURN_BOOL(FPzero(l1->A)); - PG_RETURN_BOOL(FPeq(float8_div(float8_mul(l1->A, l2->B), - float8_mul(l1->B, l2->A)), -1.0)); + PG_RETURN_BOOL(FPeq(float8_div(float8_mul(l1->A, l2->A), + float8_mul(l1->B, l2->B)), -1.0)); } Datum @@ -1133,6 +1148,20 @@ line_eq(PG_FUNCTION_ARGS) *---------------------------------------------------------*/ /* + * Return slope of the line + */ +static inline float8 +line_sl(LINE *line) +{ + if (FPzero(line->A)) + return 0.0; + if (FPzero(line->B)) + return DBL_MAX; + return float8_div(line->A, -line->B); +} + + +/* * Return inverse slope of the line */ static inline float8 @@ -1154,16 +1183,21 @@ line_distance(PG_FUNCTION_ARGS) { LINE *l1 = PG_GETARG_LINE_P(0); LINE *l2 = PG_GETARG_LINE_P(1); - float8 result; - Point tmp; + float8 ratio; if (line_interpt_line(NULL, l1, l2)) /* intersecting? */ PG_RETURN_FLOAT8(0.0); - if (FPzero(l1->B)) /* vertical? */ - PG_RETURN_FLOAT8(fabs(float8_mi(l1->C, l2->C))); - point_construct(&tmp, 0.0, l1->C); - result = line_closept_point(NULL, l2, &tmp); - PG_RETURN_FLOAT8(result); + + if (!FPzero(l1->A) && !isnan(l1->A) && !FPzero(l2->A) && !isnan(l2->A)) + ratio = float8_div(l1->A, l2->A); + else if (!FPzero(l1->B) && !isnan(l1->B) && !FPzero(l2->B) && !isnan(l2->B)) + ratio = float8_div(l1->B, l2->B); + else + ratio = 1.0; + + PG_RETURN_FLOAT8(float8_div(fabs(float8_mi(l1->C, + float8_mul(ratio, l2->C))), + HYPOT(l1->A, l1->B))); } /* line_interpt() @@ -1204,27 +1238,30 @@ line_interpt_line(Point *result, LINE *l1, LINE *l2) float8 x, y; - if (FPzero(l1->B)) /* l1 vertical? */ + if (!FPzero(l1->B)) { - if (FPzero(l2->B)) /* l2 vertical? */ + if (FPeq(l2->A, float8_mul(l1->A, float8_div(l2->B, l1->B)))) return false; - x = l1->C; - y = float8_pl(float8_mul(l2->A, x), l2->C); - } - else if (FPzero(l2->B)) /* l2 vertical? */ - { - x = l2->C; - y = float8_pl(float8_mul(l1->A, x), l1->C); + x = float8_div(float8_mi(float8_mul(l1->B, l2->C), + float8_mul(l2->B, l1->C)), + float8_mi(float8_mul(l1->A, l2->B), + float8_mul(l2->A, l1->B))); + y = float8_div(-float8_pl(float8_mul(l1->A, x), l1->C), l1->B); } - else + else if (!FPzero(l2->B)) { - if (FPeq(l2->A, float8_mul(l1->A, float8_div(l2->B, l1->B)))) + if (FPeq(l1->A, float8_mul(l2->A, float8_div(l1->B, l2->B)))) return false; - x = float8_div(float8_mi(l1->C, l2->C), float8_mi(l2->A, l1->A)); - y = float8_pl(float8_mul(l1->A, x), l1->C); + x = float8_div(float8_mi(float8_mul(l2->B, l1->C), + float8_mul(l1->B, l2->C)), + float8_mi(float8_mul(l2->A, l1->B), + float8_mul(l1->A, l2->B))); + y = float8_div(-float8_pl(float8_mul(l2->A, x), l2->C), l2->B); } + else + return false; if (result != NULL) point_construct(result, x, y); @@ -2344,16 +2381,8 @@ dist_sl(PG_FUNCTION_ARGS) { LSEG *lseg = PG_GETARG_LSEG_P(0); LINE *line = PG_GETARG_LINE_P(1); - float8 result; - - if (lseg_interpt_line(NULL, lseg, line)) - result = 0.0; - else - /* XXX shouldn't we take the min not max? */ - result = float8_max(line_closept_point(NULL, line, &lseg->p[0]), - line_closept_point(NULL, line, &lseg->p[1])); - PG_RETURN_FLOAT8(result); + PG_RETURN_FLOAT8(lseg_closept_line(NULL, lseg, line)); } /* @@ -2541,19 +2570,26 @@ lseg_interpt_line(Point *result, LSEG *lseg, LINE *line) static float8 line_closept_point(Point *result, LINE *line, Point *point) { - bool retval; - Point closept; + Point closept; LINE tmp; - /* We drop a perpendicular to find the intersection point. */ + /* + * We drop a perpendicular to find the intersection point. Ordinarily + * we should always find it, but that can fail in the presence of NaN + * coordinates, and perhaps even from simple roundoff issues. + */ line_construct(&tmp, point, line_invsl(line)); - retval = line_interpt_line(&closept, line, &tmp); - Assert(retval); /* XXX: We need something better. */ + if (!line_interpt_line(&closept, &tmp, line)) + { + if (result != NULL) + *result = *point; + + return get_float8_nan(); + } if (result != NULL) *result = closept; - /* Then we calculate the distance between the points. */ return point_dt(&closept, point); } @@ -2627,27 +2663,38 @@ lseg_closept_lseg(Point *result, LSEG *l1, LSEG *l2) float8 dist, d; - d = lseg_closept_point(NULL, l1, &l2->p[0]); - dist = d; - if (result != NULL) - *result = l2->p[0]; + /* First, we handle the case when the line segments are intersecting. */ + if (lseg_interpt_lseg(result, l1, l2)) + return 0.0; - d = lseg_closept_point(NULL, l1, &l2->p[1]); + /* + * Then, we find the closest points from the endpoints of the second + * line segment, and keep the closest one. + */ + dist = lseg_closept_point(result, l1, &l2->p[0]); + d = lseg_closept_point(&point, l1, &l2->p[1]); if (float8_lt(d, dist)) { dist = d; if (result != NULL) - *result = l2->p[1]; + *result = point; } - if (float8_lt(lseg_closept_point(&point, l2, &l1->p[0]), dist)) - d = lseg_closept_point(result, l1, &point); - - if (float8_lt(lseg_closept_point(&point, l2, &l1->p[1]), dist)) - d = lseg_closept_point(result, l1, &point); - + /* The closest point can still be one of the endpoints, so we test them. */ + d = lseg_closept_point(NULL, l2, &l1->p[0]); + if (float8_lt(d, dist)) + { + dist = d; + if (result != NULL) + *result = l1->p[0]; + } + d = lseg_closept_point(NULL, l2, &l1->p[1]); if (float8_lt(d, dist)) + { dist = d; + if (result != NULL) + *result = l1->p[1]; + } return dist; } @@ -2659,6 +2706,9 @@ close_lseg(PG_FUNCTION_ARGS) LSEG *l2 = PG_GETARG_LSEG_P(1); Point *result; + if (lseg_sl(l1) == lseg_sl(l2)) + PG_RETURN_NULL(); + result = (Point *) palloc(sizeof(Point)); if (isnan(lseg_closept_lseg(result, l2, l1))) @@ -2833,6 +2883,9 @@ close_ls(PG_FUNCTION_ARGS) LSEG *lseg = PG_GETARG_LSEG_P(1); Point *result; + if (lseg_sl(lseg) == line_sl(line)) + PG_RETURN_NULL(); + result = (Point *) palloc(sizeof(Point)); if (isnan(lseg_closept_line(result, lseg, line))) -- 2.13.6 --------------D85C8172E032009A1C02A011 Content-Type: text/x-patch; name="0003-Use-the-built-in-float-datatype-to-implement-geometr.patch" Content-Transfer-Encoding: 7bit Content-Disposition: attachment; filename*0="0003-Use-the-built-in-float-datatype-to-implement-geometr.pa"; filename*1="tch" ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
* [PATCH 2/2] Publish list of tables being repacked in shared memory @ 2026-04-07 20:29 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 940+ messages in thread From: Álvaro Herrera @ 2026-04-07 20:29 UTC (permalink / raw) Use it in autovacuum to skip processing tables that are being repacked. This is mostly to avoid repeated attempts to process such tables, which would fail due to the special deadlock checker behavior for repack. Author: Álvaro Herrera <alvherre@kurilemu.de> Discussion: https://postgr.es/m/202604072027.gazfa2zt2l2j@alvherre.pgsql --- src/backend/commands/repack.c | 195 ++++++++++++++++-- src/backend/postmaster/autovacuum.c | 20 ++ .../utils/activity/wait_event_names.txt | 1 + src/include/commands/repack.h | 2 + src/include/storage/lwlocklist.h | 2 +- src/include/storage/subsystemlist.h | 1 + src/tools/pgindent/typedefs.list | 3 + 7 files changed, 210 insertions(+), 14 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index a5f5df77291..ee7072dce6a 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -63,9 +63,11 @@ #include "optimizer/optimizer.h" #include "pgstat.h" #include "storage/bufmgr.h" +#include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" #include "storage/proc.h" +#include "storage/subsystems.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -79,6 +81,32 @@ #include "utils/syscache.h" #include "utils/wait_event_types.h" + +/* Shared memory layout for REPACK */ +typedef struct RepackWorkerInfo +{ + bool ri_in_use; + pid_t ri_backendpid; + Oid ri_dbid; + Oid ri_relid; + Oid ri_toastrelid; +} RepackWorkerInfo; + +typedef struct +{ + bool re_useless; + RepackWorkerInfo re_workerinfo[FLEXIBLE_ARRAY_MEMBER]; +} RepackShmemStruct; + +static RepackShmemStruct *RepackShmem; + +typedef struct RepackCleanupContext +{ + bool concurrent; + int workerindex; +} RepackCleanupContext; + + /* * This struct is used to pass around the information on tables to be * clustered. We need this so we can make a list of them when invoked without @@ -90,6 +118,7 @@ typedef struct Oid indexOid; } RelToCluster; + /* * The first file exported by the decoding worker must contain a snapshot, the * following ones contain the data changes. @@ -166,6 +195,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, MemoryContext permcxt); static bool repack_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); +static void RepackCleanup(RepackCleanupContext *context); +static void RepackCleanupCb(int code, Datum arg); +static void RepackShmemRequest(void *arg); +static void RepackShmemInit(void *arg); static void apply_concurrent_changes(BufFile *file, ChangeContext *chgcxt); static void apply_concurrent_insert(Relation rel, TupleTableSlot *slot, @@ -210,6 +243,11 @@ static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); +const ShmemCallbacks RepackShmemCallbacks = { + .request_fn = RepackShmemRequest, + .init_fn = RepackShmemInit, +}; + /* * The repack code allows for processing multiple tables at once. Because * of this, we cannot just run everything on a single transaction, or we @@ -514,6 +552,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid tableOid = RelationGetRelid(OldHeap); Relation index; LOCKMODE lmode; + RepackCleanupContext context; Oid save_userid; int save_sec_context; int save_nestlevel; @@ -660,24 +699,43 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, TransferPredicateLocksToHeapRelation(OldHeap); /* rebuild_relation does all the dirty work */ - PG_TRY(); - { - rebuild_relation(OldHeap, index, verbose, ident_idx); - } - PG_FINALLY(); + context.concurrent = concurrent; + + PG_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); { if (concurrent) { - /* - * Since during normal operation the worker was already asked to - * exit, stopping it explicitly is especially important on ERROR. - * However it still seems a good practice to make sure that the - * worker never survives the REPACK command. - */ - stop_repack_decoding_worker(); + bool freefound = false; + + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *worker; + + if (RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + freefound = true; + worker = &RepackShmem->re_workerinfo[i]; + context.workerindex = i; + + worker->ri_in_use = true; + worker->ri_backendpid = MyProcPid; + worker->ri_dbid = MyDatabaseId; + worker->ri_relid = RelationGetRelid(OldHeap); + worker->ri_toastrelid = OldHeap->rd_rel->reltoastrelid; + break; + } + if (!freefound) + elog(ERROR, "could not find free repack entry"); + LWLockRelease(RepackLock); } + + rebuild_relation(OldHeap, index, verbose, ident_idx); } - PG_END_TRY(); + PG_END_ENSURE_ERROR_CLEANUP(RepackCleanupCb, PointerGetDatum(&context)); + + RepackCleanup(&context); /* rebuild_relation closes OldHeap, and index if valid */ @@ -691,6 +749,117 @@ out: pgstat_progress_end_command(); } +/* + * Return whether any backend is running concurrent REPACK on the given table + * (which could be a toast table). + */ +bool +is_table_under_repack(Oid databaseId, Oid relid) +{ + bool retval = false; + + LWLockAcquire(RepackLock, LW_SHARED); + for (int i = 0; i < max_repack_replication_slots; i++) + { + RepackWorkerInfo *rworker; + + if (!RepackShmem->re_workerinfo[i].ri_in_use) + continue; + + rworker = &RepackShmem->re_workerinfo[i]; + if (rworker->ri_dbid == MyDatabaseId && + (rworker->ri_relid == relid || + rworker->ri_toastrelid == relid)) + retval = true; + } + LWLockRelease(RepackLock); + + return retval; +} + +/* + * Remove ourselves from the workerinfo array. + */ +static void +RepackCleanup(RepackCleanupContext *context) +{ + if (context->concurrent) + { + RepackWorkerInfo *worker; + + /* + * The worker would normally terminate on its own when the work is + * done, but make sure we signal it just in case. + */ + stop_repack_decoding_worker(); + + /* + * also, make sure we stop advertising the relation we were repacking, + * so that autovacuum reverts to handling it normally. + */ + LWLockAcquire(RepackLock, LW_EXCLUSIVE); + + worker = &RepackShmem->re_workerinfo[context->workerindex]; + Assert(worker->ri_backendpid == MyProcPid); + worker->ri_in_use = false; + worker->ri_backendpid = 0; + worker->ri_dbid = InvalidOid; + worker->ri_relid = InvalidOid; + worker->ri_toastrelid = InvalidOid; + LWLockRelease(RepackLock); + } +} + +/* + * RepackCleanup wrapped as an on_shmem_exit callback function + */ +static void +RepackCleanupCb(int code, Datum arg) +{ + RepackCleanup((RepackCleanupContext *) DatumGetPointer(arg)); +} + +/* + * RepackShmemRequest + * Register shared memory space needed for repack + */ +static void +RepackShmemRequest(void *arg) +{ + Size size; + + /* + * Need the fixed struct and the array of RepackWorkerInfo. + */ + size = sizeof(RepackShmemStruct); + size = MAXALIGN(size); + size = add_size(size, mul_size(max_repack_replication_slots, + sizeof(RepackWorkerInfo))); + + ShmemRequestStruct(.name = "Repack Data", + .size = size, + .ptr = (void **) &RepackShmem, + ); +} + +static void +RepackShmemInit(void *arg) +{ + RepackWorkerInfo *reinfo; + + reinfo = (RepackWorkerInfo *) ((char *) RepackShmem + + MAXALIGN(sizeof(RepackShmemStruct))); + + for (int i = 0; i < max_repack_replication_slots; i++) + { + reinfo[i].ri_in_use = false; + reinfo[i].ri_backendpid = 0; + reinfo[i].ri_dbid = InvalidOid; + reinfo[i].ri_relid = InvalidOid; + reinfo[i].ri_toastrelid = InvalidOid; + } +} + /* * Check if the table (and its index) still meets the requirements of * cluster_rel(). diff --git a/src/backend/postmaster/autovacuum.c b/src/backend/postmaster/autovacuum.c index bd626a16363..080c64ea3c8 100644 --- a/src/backend/postmaster/autovacuum.c +++ b/src/backend/postmaster/autovacuum.c @@ -78,6 +78,7 @@ #include "catalog/namespace.h" #include "catalog/pg_database.h" #include "catalog/pg_namespace.h" +#include "commands/repack.h" #include "commands/vacuum.h" #include "common/int.h" #include "funcapi.h" @@ -2422,6 +2423,25 @@ do_autovacuum(void) } } LWLockRelease(AutovacuumLock); + + /* + * Similarly, if the table is being processed by concurrent repack, + * skip it (but make a note of that). We wouldn't be able to acquire + * its lock anyway. + */ + if (!skipit) + { + MemoryContextSwitchTo(PortalContext); + + skipit = is_table_under_repack(MyDatabaseId, relid); + if (skipit) + ereport(LOG, + errmsg("skipping table \"%s.%s.%s\" because it's being repacked in concurrent mode", + get_database_name(MyDatabaseId), + get_namespace_name(get_rel_namespace(relid)), + get_rel_name(relid))); + } + if (skipit) { LWLockRelease(AutovacuumScheduleLock); diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 7bda5298558..e206304f204 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -332,6 +332,7 @@ SInvalWrite "Waiting to add a message to the shared catalog invalidation queue." WALBufMapping "Waiting to replace a page in WAL buffers." WALWrite "Waiting for WAL buffers to be written to disk." ControlFile "Waiting to read or update the <filename>pg_control</filename> file or create a new WAL file." +Repack "Waiting to read or update tables in process by concurrent repack." MultiXactGen "Waiting to read or update shared multixact state." RelCacheInit "Waiting to read or update a <filename>pg_internal.init</filename> relation cache initialization file." CheckpointerComm "Waiting to manage fsync requests." diff --git a/src/include/commands/repack.h b/src/include/commands/repack.h index fd16e74b179..be7d38b5fae 100644 --- a/src/include/commands/repack.h +++ b/src/include/commands/repack.h @@ -42,6 +42,8 @@ extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, ClusterParams *params, bool isTopLevel); +extern bool is_table_under_repack(Oid databaseId, Oid relid); + extern void check_index_is_clusterable(Relation OldHeap, Oid indexOid, LOCKMODE lockmode); extern void mark_index_clustered(Relation rel, Oid indexOid, bool is_internal); diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index af8553bcb6c..3f08f4a15d4 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -41,7 +41,7 @@ PG_LWLOCK(6, SInvalWrite) PG_LWLOCK(7, WALBufMapping) PG_LWLOCK(8, WALWrite) PG_LWLOCK(9, ControlFile) -/* 10 was CheckpointLock */ +PG_LWLOCK(10, Repack) /* 11 was XactSLRULock */ /* 12 was SubtransSLRULock */ PG_LWLOCK(13, MultiXactGen) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..4e683b8b0a8 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -72,6 +72,7 @@ PG_SHMEM_SUBSYSTEM(WalSummarizerShmemCallbacks) PG_SHMEM_SUBSYSTEM(PgArchShmemCallbacks) PG_SHMEM_SUBSYSTEM(ApplyLauncherShmemCallbacks) PG_SHMEM_SUBSYSTEM(SlotSyncShmemCallbacks) +PG_SHMEM_SUBSYSTEM(RepackShmemCallbacks) /* other modules that need some shared memory space */ PG_SHMEM_SUBSYSTEM(BTreeShmemCallbacks) diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 637c669a146..d019e03aaf1 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2639,9 +2639,12 @@ ReorderBufferTupleCidEnt ReorderBufferTupleCidKey ReorderBufferUpdateProgressTxnCB ReorderTuple +RepackCleanupContext RepackCommand RepackDecodingState +RepackShmemStruct RepackStmt +RepackWorkerInfo ReparameterizeForeignPathByChild_function ReplOriginId ReplOriginXactState -- 2.47.3 --kdrcpfmkbkc4lqhu-- ^ permalink raw reply [nested|flat] 940+ messages in thread
end of thread, other threads:[~2026-04-07 20:29 UTC | newest] Thread overview: 940+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2018-07-26 13:43 [PATCH 4/6] Fix obvious problems around the line datatype Tomas Vondra <tomas@2ndquadrant.com> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de> 2026-04-07 20:29 [PATCH 2/2] Publish list of tables being repacked in shared memory Álvaro Herrera <alvherre@kurilemu.de>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox